#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
Created on Fri Mar 27 19:38:27 2020

@author: julianduggan
"""

## import packages
import random
import time
import pandas as pd 
import glob
import os
import shutil
from selenium import webdriver
from selenium.webdriver import ActionChains
from selenium.webdriver.chrome.options import Options
from datetime import datetime

## directories - UPDATE THESE PATHS TO RUN
datasavepath = "//zip/output"
pdfsavepath = "//zip/output/pdfs"
downloadfolderpath = "/path/to/your/default/download/folder"
chromedriverpath = "/path/to/your/chromedriver/executable/file"
today=datetime.today().strftime("%Y%m%d")

## set misc. options 
pd.options.display.max_colwidth = 400

chrome_options = Options()
chrome_options.add_experimental_option('prefs',  {
    "download.default_directory": pdfsavepath,
    "download.prompt_for_download": False,
    "download.directory_upgrade": True,
    "plugins.always_open_pdf_externally": True
    }
)

## define basic functions
big = round(random.gauss(10, .1), 5)
med = round(random.gauss(7, .1), 5)
small = round(random.gauss(1, .1), 5)

def small(): 
    small_var = max(round(random.gauss(.1, .02), 5), .05)
    sleep_small = max(round(random.gauss(1, small_var), 5), 0)
    time.sleep(sleep_small)
def med(): 
    med_var = max(round(random.gauss(.3, .1), 5), .05)
    sleep_med = max(round(random.gauss(5, med_var), 5), 0)
    time.sleep(sleep_med)
def big(): 
    big_var = max(round(random.gauss(.7, .2), 5), .05)
    sleep_big = max(round(random.gauss(10, big_var), 5), 0)
    time.sleep(sleep_big)
def huge(): 
    huge_var = max(round(random.gauss(3, 1), 5), .05)   
    sleep_huge = max(round(random.gauss(60, huge_var), 5), 0)
    time.sleep(sleep_huge)
    
## set up webdriver 
driver = webdriver.Chrome(chromedriverpath + "/chromedriver")
pdfdriver= webdriver.Chrome(chromedriverpath + "/chromedriver", options=chrome_options)
actions = ActionChains(driver)
driver.implicitly_wait(5)

## get latest project list
driver.get('https://projects.worldbank.org/content/dam/documents/projects/all/all.xlsx')
 
download_complete = time.time()-os.path.getctime(downloadfolderpath)
while download_complete < 3:
    time.sleep(.5)
    download_complete = time.time()-os.path.getctime(downloadfolderpath)
time.sleep(5)
proj_file = max(glob.glob(downloadfolderpath+'/*'), key=os.path.getctime)
shutil.copyfile(proj_file, datasavepath + "/raw pipeline/all_" + today + ".xlsx") 
os.remove(proj_file)

## clean list of pipeline projects
all_proj = pd.read_excel(datasavepath + "/raw pipeline/all_" + today + ".xlsx")
if len(pd.Series.unique(all_proj['id'])) != len(all_proj['id']):
    raise Exception('Unique project ID is no longer a unique ID. Investigate!')
pipeline = all_proj[all_proj["status"].isin(["Pipeline"])]
if min(pd.isnull(pipeline['boardapprovaldate'])) == 0: 
    raise Exception('Some of these pipeline projects have board approval dates. Investigate!')
pipeline=pipeline.assign(scrapedate = '')
pipeline=pipeline.assign(docnum = '')
pipeline=pipeline.assign(pidnum = '')   
pipeline=pipeline.assign(doclinks = '')
pipeline=pipeline.assign(pidlinks = '')   

already_processed=pd.read_csv(str(glob.glob(datasavepath + "/intermediate pipeline/pipeline_all.csv")[0]))
if len(already_processed[pd.Series.isnull(already_processed['docnum'])]) != 0:
    raise Exception("Some ealrier IDs were not processed properly. Investigate!")
pipeline=pipeline[~pipeline.id.isin(already_processed.id)]

## scrape pdfs for pipeline projects
l = str(len(pipeline))
counter = 1
for pid in pipeline['id']:
    print("Now searching for " + pid + ". Number " + str(counter) + " of " + l + ".")
    
    # scrape list of URLs
    url="http://documents.worldbank.org/curated/en/docsearch?query=Project%20Information%20Document%20"+pid
    driver.get(url)
    titles=driver.find_elements_by_class_name('title')
    big()

    if len(titles) == 0: 
        pipeline.loc[pipeline['id'] == pid, 'docnum'] = 0
        pipeline.loc[pipeline['id'] == pid, 'pidnum'] = 0  
 
    # process list of possible URLs, identifying promising candidates
    links = list()
    if len(titles) > 0: 
        for t in titles: 
            link = str(t.get_attribute('href'))
            if (link != "None" and not(link in links)): 
                links.append(link)
    print(links)       
    
    if len(links) == 0:
          pipeline.loc[pipeline['id'] == pid, 'docnum'] = 0  
          pipeline.loc[pipeline['id'] == pid, 'pidnum'] = 0
    if len(links) > 0: 
        pipeline.loc[pipeline['id'] == pid, 'docnum'] = len(links)
        pipeline.loc[pipeline['id'] == pid, 'doclinks'] = ';;;; '.join(links)
        
        pidlinks = [s for s in links if (pid and "Project" and "Information" and "Document") in s]
        print(pidlinks)
        if len(pidlinks) == 0:
           pipeline.loc[pipeline['id'] == pid, 'pidnum'] = 0  
        if len(pidlinks) > 0: 
            pipeline.loc[pipeline['id'] == pid, 'pidnum'] = len(pidlinks)              
            pipeline.loc[pipeline['id'] == pid, 'pidlinks'] = ';;;; '.join(pidlinks)
    
    # visit each candidate URL to download PID pdf 
            pidindex = 1
            for plink in pidlinks: 
                driver.get(plink)
                pdflink=str(driver.find_element_by_partial_link_text('Official PDF').get_attribute('href'))
                pdfdriver.get(pdflink)
                
                download_complete = time.time()-os.path.getctime(pdfsavepath)
                while download_complete < 3:
                    time.sleep(.5)
                    download_complete = time.time()-os.path.getctime(pdfsavepath)
                big()

                pid_document=max(glob.glob(pdfsavepath+'/*'), key=os.path.getctime)
                if len(pidlinks) > 1:
                    suffix = "_" + str(pidindex)
                if len(pidlinks)==1:
                    suffix = ""
                os.rename(pid_document, pdfsavepath + "/" + pid + suffix + ".pdf")
                
                pidindex = pidindex+1
    
    counter = counter + 1
driver.close()
pdfdriver.close()

## save new dataset with scraped info
last_chunk = max(glob.glob(datasavepath+"/intermediate pipeline/pipeline_2*"), key=os.path.getctime)
nextindex=int(last_chunk.split("/")[len(last_chunk.split("/"))-1].split("_")[2].split(".")[0])+1
pipeline.to_csv(datasavepath + '/intermediate pipeline/pipeline_'+today+'_'+str(nextindex)+'.csv', encoding = 'utf-8')

clean_pipeline=pd.DataFrame()
processed_chunks=glob.glob(datasavepath + "/intermediate pipeline/pipeline_2*")
for c in processed_chunks:
    chunk=pd.read_csv(c)
    chunk=chunk[chunk.docnum.notnull()]
    clean_pipeline=clean_pipeline.append(chunk)
clean_pipeline.to_csv(datasavepath + '/intermediate pipeline/pipeline_all.csv', encoding = 'utf-8')
